agora inbox for pgsql-hackers@postgresql.org  
help / color / mirror / Atom feed
[PATCH v11 1/9] Allow CREATE INDEX CONCURRENTLY on partitioned table
500+ messages / 2 participants
[nested] [flat]

* [PATCH v11 1/9] Allow CREATE INDEX CONCURRENTLY on partitioned table
@ 2020-06-06 22:42  Justin Pryzby <pryzbyj@telsasoft.com>
  0 siblings, 0 replies; 500+ messages in thread

From: Justin Pryzby @ 2020-06-06 22:42 UTC (permalink / raw)

Note, this effectively reverts 050098b14, so take care to not reintroduce the
bug it fixed.

XXX: does pgstat_progress_update_param() break other commands progress ?
---
 doc/src/sgml/ref/create_index.sgml     |   9 --
 src/backend/commands/indexcmds.c       | 141 ++++++++++++++++++-------
 src/test/regress/expected/indexing.out |  60 ++++++++++-
 src/test/regress/sql/indexing.sql      |  18 +++-
 4 files changed, 172 insertions(+), 56 deletions(-)

diff --git a/doc/src/sgml/ref/create_index.sgml b/doc/src/sgml/ref/create_index.sgml
index 29dee5689e..bd4431a3ce 100644
--- a/doc/src/sgml/ref/create_index.sgml
+++ b/doc/src/sgml/ref/create_index.sgml
@@ -661,15 +661,6 @@ Indexes:
     cannot.
    </para>
 
-   <para>
-    Concurrent builds for indexes on partitioned tables are currently not
-    supported.  However, you may concurrently build the index on each
-    partition individually and then finally create the partitioned index
-    non-concurrently in order to reduce the time where writes to the
-    partitioned table will be locked out.  In this case, building the
-    partitioned index is a metadata only operation.
-   </para>
-
   </refsect2>
  </refsect1>
 
diff --git a/src/backend/commands/indexcmds.c b/src/backend/commands/indexcmds.c
index ca24620fd0..76219381c1 100644
--- a/src/backend/commands/indexcmds.c
+++ b/src/backend/commands/indexcmds.c
@@ -68,6 +68,7 @@
 
 
 /* non-export function prototypes */
+static void reindex_invalid_child_indexes(Oid indexRelationId);
 static bool CompareOpclassOptions(Datum *opts1, Datum *opts2, int natts);
 static void CheckPredicate(Expr *predicate);
 static void ComputeIndexAttrs(IndexInfo *indexInfo,
@@ -671,17 +672,6 @@ DefineIndex(Oid relationId,
 	partitioned = rel->rd_rel->relkind == RELKIND_PARTITIONED_TABLE;
 	if (partitioned)
 	{
-		/*
-		 * Note: we check 'stmt->concurrent' rather than 'concurrent', so that
-		 * the error is thrown also for temporary tables.  Seems better to be
-		 * consistent, even though we could do it on temporary table because
-		 * we're not actually doing it concurrently.
-		 */
-		if (stmt->concurrent)
-			ereport(ERROR,
-					(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
-					 errmsg("cannot create index on partitioned table \"%s\" concurrently",
-							RelationGetRelationName(rel))));
 		if (stmt->excludeOpNames)
 			ereport(ERROR,
 					(errcode(ERRCODE_FEATURE_NOT_SUPPORTED),
@@ -1119,6 +1109,11 @@ DefineIndex(Oid relationId,
 		if (pd->nparts != 0)
 			flags |= INDEX_CREATE_INVALID;
 	}
+	else if (concurrent && OidIsValid(parentIndexId))
+	{
+		/* If concurrent, initially build index partitions as "invalid" */
+		flags |= INDEX_CREATE_INVALID;
+	}
 
 	if (stmt->deferrable)
 		constr_flags |= INDEX_CONSTR_CREATE_DEFERRABLE;
@@ -1171,6 +1166,14 @@ DefineIndex(Oid relationId,
 		 */
 		if (!stmt->relation || stmt->relation->inh)
 		{
+			/*
+			 * Need to close the relation before recursing into children, so
+			 * copy needed data into a longlived context.
+			 */
+
+			MemoryContext	ind_context = AllocSetContextCreate(PortalContext, "CREATE INDEX",
+					ALLOCSET_DEFAULT_SIZES);
+			MemoryContext	oldcontext = MemoryContextSwitchTo(ind_context);
 			PartitionDesc partdesc = RelationGetPartitionDesc(rel);
 			int			nparts = partdesc->nparts;
 			Oid		   *part_oids = palloc(sizeof(Oid) * nparts);
@@ -1178,12 +1181,15 @@ DefineIndex(Oid relationId,
 			TupleDesc	parentDesc;
 			Oid		   *opfamOids;
 
+			// If concurrent, maybe this should be done after excluding indexes which already exist ?
 			pgstat_progress_update_param(PROGRESS_CREATEIDX_PARTITIONS_TOTAL,
 										 nparts);
 
 			memcpy(part_oids, partdesc->oids, sizeof(Oid) * nparts);
+			parentDesc = CreateTupleDescCopy(RelationGetDescr(rel));
+			table_close(rel, NoLock);
+			MemoryContextSwitchTo(oldcontext);
 
-			parentDesc = RelationGetDescr(rel);
 			opfamOids = palloc(sizeof(Oid) * numberOfKeyAttributes);
 			for (i = 0; i < numberOfKeyAttributes; i++)
 				opfamOids[i] = get_opclass_family(classObjectId[i]);
@@ -1226,10 +1232,12 @@ DefineIndex(Oid relationId,
 					continue;
 				}
 
+				oldcontext = MemoryContextSwitchTo(ind_context);
 				childidxs = RelationGetIndexList(childrel);
 				attmap =
 					build_attrmap_by_name(RelationGetDescr(childrel),
 										  parentDesc);
+				MemoryContextSwitchTo(oldcontext);
 
 				foreach(cell, childidxs)
 				{
@@ -1300,10 +1308,14 @@ DefineIndex(Oid relationId,
 				 */
 				if (!found)
 				{
-					IndexStmt  *childStmt = copyObject(stmt);
+					IndexStmt  *childStmt;
 					bool		found_whole_row;
 					ListCell   *lc;
 
+					oldcontext = MemoryContextSwitchTo(ind_context);
+					childStmt = copyObject(stmt);
+					MemoryContextSwitchTo(oldcontext);
+
 					/*
 					 * We can't use the same index name for the child index,
 					 * so clear idxname to let the recursive invocation choose
@@ -1355,10 +1367,18 @@ DefineIndex(Oid relationId,
 								createdConstraintId,
 								is_alter_table, check_rights, check_not_in_use,
 								skip_build, quiet);
+					if (concurrent)
+					{
+						PopActiveSnapshot();
+						PushActiveSnapshot(GetTransactionSnapshot());
+						invalidate_parent = true;
+					}
 				}
 
-				pgstat_progress_update_param(PROGRESS_CREATEIDX_PARTITIONS_DONE,
-											 i + 1);
+				/* For concurrent build, this is a catalog-only stage */
+				if (!concurrent)
+					pgstat_progress_update_param(PROGRESS_CREATEIDX_PARTITIONS_DONE,
+												 i + 1);
 				free_attrmap(attmap);
 			}
 
@@ -1368,51 +1388,42 @@ DefineIndex(Oid relationId,
 			 * invalid, this is incorrect, so update our row to invalid too.
 			 */
 			if (invalidate_parent)
-			{
-				Relation	pg_index = table_open(IndexRelationId, RowExclusiveLock);
-				HeapTuple	tup,
-							newtup;
-
-				tup = SearchSysCache1(INDEXRELID,
-									  ObjectIdGetDatum(indexRelationId));
-				if (!HeapTupleIsValid(tup))
-					elog(ERROR, "cache lookup failed for index %u",
-						 indexRelationId);
-				newtup = heap_copytuple(tup);
-				((Form_pg_index) GETSTRUCT(newtup))->indisvalid = false;
-				CatalogTupleUpdate(pg_index, &tup->t_self, newtup);
-				ReleaseSysCache(tup);
-				table_close(pg_index, RowExclusiveLock);
-				heap_freetuple(newtup);
-			}
-		}
+				index_set_state_flags(indexRelationId, INDEX_DROP_CLEAR_VALID);
+		} else
+			table_close(rel, NoLock);
 
 		/*
 		 * Indexes on partitioned tables are not themselves built, so we're
 		 * done here.
 		 */
-		table_close(rel, NoLock);
 		if (!OidIsValid(parentIndexId))
+		{
+			if (concurrent)
+				reindex_invalid_child_indexes(indexRelationId);
+
 			pgstat_progress_end_command();
+		}
+
 		return address;
 	}
 
-	if (!concurrent)
+	table_close(rel, NoLock);
+	if (!concurrent || OidIsValid(parentIndexId))
 	{
-		/* Close the heap and we're done, in the non-concurrent case */
-		table_close(rel, NoLock);
+		/*
+		 * We're done if this is the top-level index,
+		 * or the catalog-only phase of a partition built concurrently
+		 */
 
-		/* If this is the top-level index, we're done. */
 		if (!OidIsValid(parentIndexId))
 			pgstat_progress_end_command();
 
 		return address;
 	}
 
-	/* save lockrelid and locktag for below, then close rel */
+	/* save lockrelid and locktag for below */
 	heaprelid = rel->rd_lockInfo.lockRelId;
 	SET_LOCKTAG_RELATION(heaplocktag, heaprelid.dbId, heaprelid.relId);
-	table_close(rel, NoLock);
 
 	/*
 	 * For a concurrent build, it's important to make the catalog entries
@@ -1606,6 +1617,56 @@ DefineIndex(Oid relationId,
 	return address;
 }
 
+/* Reindex invalid child indexes created earlier */
+static void
+reindex_invalid_child_indexes(Oid indexRelationId)
+{
+	ListCell *lc;
+	int		npart = 0;
+
+	MemoryContext	ind_context = AllocSetContextCreate(PortalContext, "CREATE INDEX",
+			ALLOCSET_DEFAULT_SIZES);
+	MemoryContext	oldcontext;
+	List		*childs = find_inheritance_children(indexRelationId, ShareLock);
+	List		*partitions = NIL;
+
+	PreventInTransactionBlock(true, "REINDEX INDEX");
+
+	foreach (lc, childs)
+	{
+		Oid			partoid = lfirst_oid(lc);
+
+		pgstat_progress_update_param(PROGRESS_CREATEIDX_PARTITIONS_DONE,
+									 npart++);
+
+		if (get_index_isvalid(partoid) ||
+				!RELKIND_HAS_STORAGE(get_rel_relkind(partoid)))
+			continue;
+
+		/* Save partition OID */
+		oldcontext = MemoryContextSwitchTo(ind_context);
+		partitions = lappend_oid(partitions, partoid);
+		MemoryContextSwitchTo(oldcontext);
+	}
+
+	/*
+	 * Process each partition listed in a separate transaction.  Note that
+	 * this commits and then starts a new transaction immediately.
+	 */
+	ReindexMultipleInternal(partitions, REINDEXOPT_CONCURRENTLY);
+
+	/*
+	 * CIC needs to mark a partitioned index as VALID, which itself
+	 * requires setting READY, which is unset for CIC (even though
+	 * it's meaningless for an index without storage).
+	 * This must be done only while holding a lock which precludes adding
+	 * partitions.
+	 * See also: validatePartitionedIndex().
+	 */
+	index_set_state_flags(indexRelationId, INDEX_CREATE_SET_READY);
+	CommandCounterIncrement();
+	index_set_state_flags(indexRelationId, INDEX_CREATE_SET_VALID);
+}
 
 /*
  * CheckMutability
diff --git a/src/test/regress/expected/indexing.out b/src/test/regress/expected/indexing.out
index c93f4470c9..f04abc6897 100644
--- a/src/test/regress/expected/indexing.out
+++ b/src/test/regress/expected/indexing.out
@@ -50,11 +50,63 @@ select relname, relkind, relhassubclass, inhparent::regclass
 (8 rows)
 
 drop table idxpart;
--- Some unsupported features
+-- CIC on partitioned table
 create table idxpart (a int, b int, c text) partition by range (a);
-create table idxpart1 partition of idxpart for values from (0) to (10);
-create index concurrently on idxpart (a);
-ERROR:  cannot create index on partitioned table "idxpart" concurrently
+create table idxpart1 partition of idxpart for values from (0) to (10) partition by range(a);
+create table idxpart11 partition of idxpart1 for values from (0) to (10) partition by range(a);
+create table idxpart111 partition of idxpart11 default partition by range(a);
+create table idxpart1111 partition of idxpart111 default partition by range(a);
+create table idxpart2 partition of idxpart for values from (10) to (20);
+insert into idxpart2 values(10),(10); -- not unique
+create index concurrently on idxpart (a); -- partitioned
+create index concurrently on idxpart1 (a); -- partitioned and partition
+create index concurrently on idxpart11 (a); -- partitioned and partition, with no leaves
+create index concurrently on idxpart2 (a); -- leaf
+create unique index concurrently on idxpart (a); -- partitioned, unique failure
+ERROR:  could not create unique index "idxpart2_a_idx2_ccnew"
+DETAIL:  Key (a)=(10) is duplicated.
+\d idxpart
+        Partitioned table "public.idxpart"
+ Column |  Type   | Collation | Nullable | Default 
+--------+---------+-----------+----------+---------
+ a      | integer |           |          | 
+ b      | integer |           |          | 
+ c      | text    |           |          | 
+Partition key: RANGE (a)
+Indexes:
+    "idxpart_a_idx" btree (a)
+    "idxpart_a_idx1" UNIQUE, btree (a) INVALID
+Number of partitions: 2 (Use \d+ to list them.)
+
+\d idxpart1
+        Partitioned table "public.idxpart1"
+ Column |  Type   | Collation | Nullable | Default 
+--------+---------+-----------+----------+---------
+ a      | integer |           |          | 
+ b      | integer |           |          | 
+ c      | text    |           |          | 
+Partition of: idxpart FOR VALUES FROM (0) TO (10)
+Partition key: RANGE (a)
+Indexes:
+    "idxpart1_a_idx" btree (a) INVALID
+    "idxpart1_a_idx1" btree (a)
+    "idxpart1_a_idx2" UNIQUE, btree (a) INVALID
+Number of partitions: 1 (Use \d+ to list them.)
+
+\d idxpart2
+              Table "public.idxpart2"
+ Column |  Type   | Collation | Nullable | Default 
+--------+---------+-----------+----------+---------
+ a      | integer |           |          | 
+ b      | integer |           |          | 
+ c      | text    |           |          | 
+Partition of: idxpart FOR VALUES FROM (10) TO (20)
+Indexes:
+    "idxpart2_a_idx" btree (a)
+    "idxpart2_a_idx1" btree (a)
+    "idxpart2_a_idx2" UNIQUE, btree (a) INVALID
+    "idxpart2_a_idx2_ccnew" UNIQUE, btree (a) INVALID
+
 drop table idxpart;
 -- Verify bugfix with query on indexed partitioned table with no partitions
 -- https://postgr.es/m/20180124162006.pmapfiznhgngwtjf@alvherre.pgsql
diff --git a/src/test/regress/sql/indexing.sql b/src/test/regress/sql/indexing.sql
index 42f398b67c..3d4b6e9bc9 100644
--- a/src/test/regress/sql/indexing.sql
+++ b/src/test/regress/sql/indexing.sql
@@ -29,10 +29,22 @@ select relname, relkind, relhassubclass, inhparent::regclass
 	where relname like 'idxpart%' order by relname;
 drop table idxpart;
 
--- Some unsupported features
+-- CIC on partitioned table
 create table idxpart (a int, b int, c text) partition by range (a);
-create table idxpart1 partition of idxpart for values from (0) to (10);
-create index concurrently on idxpart (a);
+create table idxpart1 partition of idxpart for values from (0) to (10) partition by range(a);
+create table idxpart11 partition of idxpart1 for values from (0) to (10) partition by range(a);
+create table idxpart111 partition of idxpart11 default partition by range(a);
+create table idxpart1111 partition of idxpart111 default partition by range(a);
+create table idxpart2 partition of idxpart for values from (10) to (20);
+insert into idxpart2 values(10),(10); -- not unique
+create index concurrently on idxpart (a); -- partitioned
+create index concurrently on idxpart1 (a); -- partitioned and partition
+create index concurrently on idxpart11 (a); -- partitioned and partition, with no leaves
+create index concurrently on idxpart2 (a); -- leaf
+create unique index concurrently on idxpart (a); -- partitioned, unique failure
+\d idxpart
+\d idxpart1
+\d idxpart2
 drop table idxpart;
 
 -- Verify bugfix with query on indexed partitioned table with no partitions
-- 
2.17.0


--cWoXeonUoKmBZSoM
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment;
 filename="v11-0002-Add-SKIPVALID-flag-for-more-integration.patch"



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread

* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21  Nathan Bossart <nathan@postgresql.org>
  0 siblings, 0 replies; 500+ messages in thread

From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)

---
 src/backend/access/heap/heapam_handler.c |  2 +-
 src/backend/access/table/tableam.c       | 59 +++++++++++-------------
 src/include/access/relscan.h             |  8 ++--
 3 files changed, 31 insertions(+), 38 deletions(-)

diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
 	if (hscan->rs_base.rs_parallel != NULL)
 	{
 		bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
-		startblock = bpscan->phs_startblock;
+		startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
 	}
 	else
 		startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
 	bpscan->base.phs_syncscan = synchronize_seqscans &&
 		!RelationUsesLocalBuffers(rel) &&
 		bpscan->phs_nblocks > NBuffers / 4;
-	SpinLockInit(&bpscan->phs_mutex);
-	bpscan->phs_startblock = InvalidBlockNumber;
-	bpscan->phs_numblock = InvalidBlockNumber;
+	pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+	pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
 	pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
 
 	return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
 	StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
 					 "pg_nextpower2_32 may be too small for non-standard BlockNumber width");
 
-	BlockNumber sync_startpage = InvalidBlockNumber;
 	BlockNumber scan_nblocks;
 
 	/* Reset the state we use for controlling allocation size. */
 	memset(pbscanwork, 0, sizeof(*pbscanwork));
 
-retry:
-	/* Grab the spinlock. */
-	SpinLockAcquire(&pbscan->phs_mutex);
-
 	/*
 	 * When the caller specified a limit on the number of blocks to scan, set
 	 * that in the ParallelBlockTableScanDesc, if it's not been done by
 	 * another worker already.
 	 */
-	if (numblocks != InvalidBlockNumber &&
-		pbscan->phs_numblock == InvalidBlockNumber)
+	if (numblocks != InvalidBlockNumber)
 	{
-		pbscan->phs_numblock = numblocks;
+		uint32		expected = InvalidBlockNumber;
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+									   numblocks);
 	}
 
 	/*
@@ -485,36 +481,35 @@ retry:
 	 * so now.  If a startblock was specified, start there, otherwise if this
 	 * is not a synchronized scan, we just start at block 0, but if it is a
 	 * synchronized scan, we must get the starting position from the
-	 * synchronized scan machinery.  We can't hold the spinlock while doing
-	 * that, though, so release the spinlock, get the information we need, and
-	 * retry.  If nobody else has initialized the scan in the meantime, we'll
-	 * fill in the value we fetched on the second time through.
+	 * synchronized scan machinery.
+	 *
+	 * If another worker initializes phs_startblock concurrently, just use
+	 * their value.
 	 */
-	if (pbscan->phs_startblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
 	{
+		BlockNumber newstartblock;
+		uint32		expected = InvalidBlockNumber;
+
 		if (startblock != InvalidBlockNumber)
-			pbscan->phs_startblock = startblock;
+			newstartblock = startblock;
 		else if (!pbscan->base.phs_syncscan)
-			pbscan->phs_startblock = 0;
-		else if (sync_startpage != InvalidBlockNumber)
-			pbscan->phs_startblock = sync_startpage;
+			newstartblock = 0;
 		else
-		{
-			SpinLockRelease(&pbscan->phs_mutex);
-			sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
-			goto retry;
-		}
+			newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+		pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+									   newstartblock);
 	}
-	SpinLockRelease(&pbscan->phs_mutex);
 
 	/*
 	 * Figure out how many blocks we're going to scan; either all of them, or
 	 * just phs_numblock's worth, if a limit has been imposed.
 	 */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * We determine the chunk size based on scan_nblocks.  First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
 	 */
 
 	/* First, figure out how many blocks we're planning on scanning */
-	if (pbscan->phs_numblock == InvalidBlockNumber)
+	if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
 		scan_nblocks = pbscan->phs_nblocks;
 	else
-		scan_nblocks = pbscan->phs_numblock;
+		scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
 
 	/*
 	 * Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
 	if (nallocated >= scan_nblocks)
 		page = InvalidBlockNumber;	/* all blocks have been allocated */
 	else
-		page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+		page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
 
 	/*
 	 * Report scan location.  Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
 		if (page != InvalidBlockNumber)
 			ss_report_location(rel, page);
 		else if (nallocated == pbscan->phs_nblocks)
-			ss_report_location(rel, pbscan->phs_startblock);
+			ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
 	}
 
 	return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
 #include "nodes/tidbitmap.h"
 #include "port/atomics.h"
 #include "storage/relfilelocator.h"
-#include "storage/spin.h"
 #include "utils/relcache.h"
 
 
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
 	ParallelTableScanDescData base;
 
 	BlockNumber phs_nblocks;	/* # blocks in relation at start of scan */
-	slock_t		phs_mutex;		/* mutual exclusion for setting startblock */
-	BlockNumber phs_startblock; /* starting block number */
-	BlockNumber phs_numblock;	/* # blocks to scan, or InvalidBlockNumber if
-								 * no limit */
+	pg_atomic_uint32 phs_startblock;	/* starting block number */
+	pg_atomic_uint32 phs_numblock;	/* # blocks to scan, or InvalidBlockNumber
+									 * if no limit */
 	pg_atomic_uint64 phs_nallocated;	/* number of blocks allocated to
 										 * workers so far. */
 }			ParallelBlockTableScanDescData;
-- 
2.50.1 (Apple Git-155)


--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
	filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch



^ permalink  raw  reply  [nested|flat] 500+ messages in thread


end of thread, other threads:[~2026-07-09 20:21 UTC | newest]

Thread overview: 500+ messages (download: mbox mbox.gz follow: Atom feed)
-- links below jump to the message on this page --
2020-06-06 22:42 [PATCH v11 1/9] Allow CREATE INDEX CONCURRENTLY on partitioned table Justin Pryzby <pryzbyj@telsasoft.com>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>

This inbox is served by agora; see mirroring instructions
for how to clone and mirror all data and code used for this inbox